agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
"""Stage four: run the scenarios against the world and say what happened.
|
|
2
|
+
|
|
3
|
+
Every scenario gets its own world. It is restored from the frozen snapshot, the scenario's own
|
|
4
|
+
setup is run against it, and it is thrown away afterwards. Nothing a scenario does can reach the
|
|
5
|
+
next one, which is what makes a result mean something on its own and makes the whole suite
|
|
6
|
+
repeatable a week later.
|
|
7
|
+
|
|
8
|
+
The shape is the same regardless of what is being tested: restore, converse, grade against the
|
|
9
|
+
state that is left behind. Where the agent actually runs is a target, so the same scenarios grade
|
|
10
|
+
a hosted agent without any of this changing.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Callable, Sequence
|
|
18
|
+
|
|
19
|
+
from ..contract import AgentContract
|
|
20
|
+
from ..catalogue import load_catalogue
|
|
21
|
+
from ..simulator import load_simulator_prompt
|
|
22
|
+
from ..scenario import Scenario
|
|
23
|
+
from ..folder import apply_setup, check_ready
|
|
24
|
+
from ..world.snapshot import restore
|
|
25
|
+
from .conversation import FINISHED, Exchange, Transcript, converse
|
|
26
|
+
from .grade import (
|
|
27
|
+
Checkpoint,
|
|
28
|
+
Result,
|
|
29
|
+
as_json,
|
|
30
|
+
checkpoints,
|
|
31
|
+
grade_sub_goals,
|
|
32
|
+
judge,
|
|
33
|
+
judge_suite_evals,
|
|
34
|
+
summarise,
|
|
35
|
+
ungraded_sub_goals,
|
|
36
|
+
)
|
|
37
|
+
from .targets import (
|
|
38
|
+
LocalAgent,
|
|
39
|
+
RepositoryChatTarget,
|
|
40
|
+
Target,
|
|
41
|
+
register_target,
|
|
42
|
+
resolve,
|
|
43
|
+
supported,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _cases(report: Any) -> list[Any]:
|
|
48
|
+
"""The per-scenario results, whichever report this is.
|
|
49
|
+
|
|
50
|
+
The runner hands back a ``SimulationReport``, whose cases are ``test_cases`` and whose
|
|
51
|
+
messages live a level down; the plugins hand back the older ``TestReport``, whose cases are
|
|
52
|
+
``results``. Reading only one of them finds nothing in the other and reports a run in which
|
|
53
|
+
nobody said anything, which is indistinguishable from an agent that ignored the person.
|
|
54
|
+
"""
|
|
55
|
+
legacy = getattr(report, "to_legacy", None)
|
|
56
|
+
if callable(legacy):
|
|
57
|
+
try:
|
|
58
|
+
report = legacy()
|
|
59
|
+
except Exception: # noqa: BLE001 - a report that will not convert is still readable
|
|
60
|
+
pass
|
|
61
|
+
return list(
|
|
62
|
+
getattr(report, "results", None) or getattr(report, "test_cases", None) or []
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def from_alk(report: Any, world, spent: float) -> Transcript:
|
|
67
|
+
"""What ALK's report says happened, in the shape the grading already reads.
|
|
68
|
+
|
|
69
|
+
Read from ``messages``, the normalised trajectory, and not from ``transcript`` — which is a
|
|
70
|
+
string, so iterating it yields characters, matches nothing, and produces a run with no turns
|
|
71
|
+
at all. The judge is then handed an empty conversation and fails every claim about what was
|
|
72
|
+
said, which arrives looking like an agent that never spoke.
|
|
73
|
+
"""
|
|
74
|
+
exchanges: list[Exchange] = []
|
|
75
|
+
for case in _cases(report):
|
|
76
|
+
for message in getattr(case, "messages", None) or []:
|
|
77
|
+
if not isinstance(message, dict):
|
|
78
|
+
continue
|
|
79
|
+
role = str(message.get("role") or "")
|
|
80
|
+
text = message.get("content") or ""
|
|
81
|
+
# Tool turns are in here too, and they are already recorded as calls. Putting them
|
|
82
|
+
# in the conversation as well would have the judge read a tool result as something
|
|
83
|
+
# the agent said.
|
|
84
|
+
if role in ("tool", "function") or not str(text).strip():
|
|
85
|
+
continue
|
|
86
|
+
exchanges.append(
|
|
87
|
+
Exchange("agent" if role == "assistant" else "customer", str(text))
|
|
88
|
+
)
|
|
89
|
+
if not exchanges and isinstance(getattr(case, "transcript", None), str):
|
|
90
|
+
# A plugin that only fills the text form still has to be readable.
|
|
91
|
+
spoken = case.transcript.strip()
|
|
92
|
+
if spoken:
|
|
93
|
+
exchanges.append(Exchange("customer", spoken))
|
|
94
|
+
return Transcript(
|
|
95
|
+
exchanges=exchanges,
|
|
96
|
+
calls=list(world.calls),
|
|
97
|
+
ended=FINISHED,
|
|
98
|
+
spent_usd=spent,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def audio_from(report: Any) -> str:
|
|
103
|
+
"""Where ALK left this run's audio, if it left any.
|
|
104
|
+
|
|
105
|
+
Artifacts are how a modality hands back what it produced, so the recording is asked of the
|
|
106
|
+
report rather than guessed at from a directory. A run with none says so.
|
|
107
|
+
"""
|
|
108
|
+
for case in _cases(report):
|
|
109
|
+
for artifact in getattr(case, "artifacts", None) or []:
|
|
110
|
+
kind = str(getattr(artifact, "type", "") or "")
|
|
111
|
+
mime = str(getattr(artifact, "mime_type", "") or "")
|
|
112
|
+
if "audio" in kind.lower() or mime.startswith("audio/"):
|
|
113
|
+
found = getattr(artifact, "path", None) or getattr(
|
|
114
|
+
artifact, "uri", None
|
|
115
|
+
)
|
|
116
|
+
if found:
|
|
117
|
+
return str(found)
|
|
118
|
+
return ""
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
RUNS = "runs.json"
|
|
122
|
+
REPORT = "report.txt"
|
|
123
|
+
|
|
124
|
+
__all__ = [
|
|
125
|
+
"Checkpoint",
|
|
126
|
+
"Exchange",
|
|
127
|
+
"LocalAgent",
|
|
128
|
+
"RepositoryChatTarget",
|
|
129
|
+
"Result",
|
|
130
|
+
"Target",
|
|
131
|
+
"Transcript",
|
|
132
|
+
"converse",
|
|
133
|
+
"register_target",
|
|
134
|
+
"run_scenario",
|
|
135
|
+
"run_suite",
|
|
136
|
+
"supported",
|
|
137
|
+
"summarise",
|
|
138
|
+
]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
async def run_scenario(
|
|
142
|
+
scenario: Scenario,
|
|
143
|
+
contract: AgentContract,
|
|
144
|
+
world_root: Path,
|
|
145
|
+
*,
|
|
146
|
+
target: str = "local",
|
|
147
|
+
model: str | None = None,
|
|
148
|
+
through_alk: bool = False,
|
|
149
|
+
on_exchange: Callable[[Exchange], Any] | None = None,
|
|
150
|
+
) -> Result:
|
|
151
|
+
"""Run one scenario in its own copy of the world and grade what it left behind."""
|
|
152
|
+
catalogue = load_catalogue(world_root)
|
|
153
|
+
world = restore(world_root)
|
|
154
|
+
try:
|
|
155
|
+
# reset() is how an environment is started in ALK: it clears the call log and
|
|
156
|
+
# publishes the tools and the starting state. Going through it keeps a generated world
|
|
157
|
+
# drivable by anything that already drives an environment.
|
|
158
|
+
world.reset()
|
|
159
|
+
applied = apply_setup(scenario, world)
|
|
160
|
+
if not applied.ok:
|
|
161
|
+
raise RuntimeError(f"the scenario's setup did not run: {applied.said}")
|
|
162
|
+
ready = check_ready(scenario, world)
|
|
163
|
+
if not ready.ok:
|
|
164
|
+
raise RuntimeError(
|
|
165
|
+
f"the world is not ready for this scenario: {ready.said}. Running it would "
|
|
166
|
+
"test us rather than the agent."
|
|
167
|
+
)
|
|
168
|
+
# The setup's calls are not the agent's.
|
|
169
|
+
world.calls = []
|
|
170
|
+
if through_alk:
|
|
171
|
+
# ALK owns the simulation and drives the world through EnvironmentAdapter; the
|
|
172
|
+
# harness only grades what it is left with. Nothing here is modality-specific,
|
|
173
|
+
# which is the point: the browser and voice runners take the same adapter.
|
|
174
|
+
from .alk import simulate
|
|
175
|
+
|
|
176
|
+
report, spent = await simulate(
|
|
177
|
+
scenario,
|
|
178
|
+
contract,
|
|
179
|
+
world,
|
|
180
|
+
model=model,
|
|
181
|
+
simulator_prompt=load_simulator_prompt(world_root),
|
|
182
|
+
)
|
|
183
|
+
transcript = from_alk(report, world, spent)
|
|
184
|
+
for exchange in transcript.exchanges:
|
|
185
|
+
if on_exchange:
|
|
186
|
+
on_exchange(exchange)
|
|
187
|
+
else:
|
|
188
|
+
agent = resolve(target)(contract, world, model=model)
|
|
189
|
+
transcript = await converse(
|
|
190
|
+
agent,
|
|
191
|
+
scenario,
|
|
192
|
+
contract,
|
|
193
|
+
world_root=world_root,
|
|
194
|
+
model=model,
|
|
195
|
+
on_exchange=on_exchange,
|
|
196
|
+
)
|
|
197
|
+
# Settled by code first. The judge is only handed the sub-goals whose catalogue entry
|
|
198
|
+
# says nothing observable decides them.
|
|
199
|
+
settled = grade_sub_goals(world, scenario, catalogue, transcript.calls)
|
|
200
|
+
ending = ", ".join(
|
|
201
|
+
f"{name}: {len(rows)} rows"
|
|
202
|
+
for name, rows in sorted(world.observe().state.items())
|
|
203
|
+
)
|
|
204
|
+
judgements, judged_cost = await judge(
|
|
205
|
+
scenario, transcript, contract, catalogue, model=model, ending=ending
|
|
206
|
+
)
|
|
207
|
+
judgements += judge_suite_evals(
|
|
208
|
+
catalogue.suite_evals, scenario, transcript, contract, ending=ending
|
|
209
|
+
)
|
|
210
|
+
return Result(
|
|
211
|
+
scenario=scenario.name,
|
|
212
|
+
tests=scenario.tests,
|
|
213
|
+
problems=[
|
|
214
|
+
f"{name} is not in this catalogue, so nothing graded it"
|
|
215
|
+
for name in ungraded_sub_goals(scenario, catalogue)
|
|
216
|
+
],
|
|
217
|
+
state_failures=[
|
|
218
|
+
f"{one.name}: {one.said}" for one in settled if not one.held
|
|
219
|
+
],
|
|
220
|
+
conduct=judgements,
|
|
221
|
+
checkpoints=checkpoints(settled, judgements),
|
|
222
|
+
crashes=[f"{call.name}: {call.error}" for call in transcript.crashed()],
|
|
223
|
+
ended=transcript.ended,
|
|
224
|
+
turns=len(transcript.exchanges),
|
|
225
|
+
calls=len(transcript.calls),
|
|
226
|
+
spent_usd=transcript.spent_usd + judged_cost,
|
|
227
|
+
transcript=transcript.spoken(),
|
|
228
|
+
exchanges=[
|
|
229
|
+
{"speaker": turn.speaker, "text": turn.text}
|
|
230
|
+
for turn in transcript.exchanges
|
|
231
|
+
],
|
|
232
|
+
actions=transcript.actions(),
|
|
233
|
+
)
|
|
234
|
+
finally:
|
|
235
|
+
world.close()
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
async def run_suite(
|
|
239
|
+
scenarios: Sequence[Scenario],
|
|
240
|
+
contract: AgentContract,
|
|
241
|
+
world_root: Path,
|
|
242
|
+
*,
|
|
243
|
+
target: str = "local",
|
|
244
|
+
model: str | None = None,
|
|
245
|
+
through_alk: bool = False,
|
|
246
|
+
out: Path | None = None,
|
|
247
|
+
on_result: Callable[[Result], Any] | None = None,
|
|
248
|
+
on_exchange: Callable[[Exchange], Any] | None = None,
|
|
249
|
+
) -> list[Result]:
|
|
250
|
+
"""Run every scenario and write the results out. One failing scenario never stops the rest."""
|
|
251
|
+
destination = Path(out or world_root)
|
|
252
|
+
results: list[Result] = []
|
|
253
|
+
for scenario in scenarios:
|
|
254
|
+
try:
|
|
255
|
+
result = await run_scenario(
|
|
256
|
+
scenario,
|
|
257
|
+
contract,
|
|
258
|
+
world_root,
|
|
259
|
+
target=target,
|
|
260
|
+
model=model,
|
|
261
|
+
through_alk=through_alk,
|
|
262
|
+
on_exchange=on_exchange,
|
|
263
|
+
)
|
|
264
|
+
except Exception as failed:
|
|
265
|
+
# A scenario that could not be run is recorded as unrunnable rather than as a
|
|
266
|
+
# failure of the agent, and the rest of the suite still runs.
|
|
267
|
+
result = Result(
|
|
268
|
+
scenario=scenario.name,
|
|
269
|
+
tests=scenario.tests,
|
|
270
|
+
crashes=[f"could not run: {type(failed).__name__}: {failed}"],
|
|
271
|
+
ended="not-run",
|
|
272
|
+
)
|
|
273
|
+
results.append(result)
|
|
274
|
+
if on_result:
|
|
275
|
+
on_result(result)
|
|
276
|
+
|
|
277
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
278
|
+
# Records for scenarios this suite did not run are kept, not clobbered. A live call and a
|
|
279
|
+
# local run write to the same file, and re-running two scenarios must not erase the third.
|
|
280
|
+
ran = {result.scenario for result in results}
|
|
281
|
+
kept = [
|
|
282
|
+
record
|
|
283
|
+
for record in load_results(destination)
|
|
284
|
+
if isinstance(record, dict) and record.get("scenario") not in ran
|
|
285
|
+
]
|
|
286
|
+
merged = kept + json.loads(as_json(results))
|
|
287
|
+
(destination / RUNS).write_text(
|
|
288
|
+
json.dumps(merged, indent=2, ensure_ascii=False), encoding="utf-8"
|
|
289
|
+
)
|
|
290
|
+
(destination / REPORT).write_text(summarise(results), encoding="utf-8")
|
|
291
|
+
return results
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def load_results(destination: Path) -> list[dict[str, Any]]:
|
|
295
|
+
path = Path(destination) / RUNS
|
|
296
|
+
return json.loads(path.read_text(encoding="utf-8")) if path.exists() else []
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
"""Running a generated world through ALK's own simulation, rather than beside it.
|
|
2
|
+
|
|
3
|
+
The whole reason a generated world subclasses ``EnvironmentAdapter`` is so the runners that
|
|
4
|
+
already exist can drive it. ``ChatEnvironment`` takes ``environment=<adapter>`` and owns the
|
|
5
|
+
synthetic user, the turn loop, the transcript and the report; the browser and voice paths take
|
|
6
|
+
the same adapter. A second loop written here would work for exactly one modality and would have
|
|
7
|
+
to be rewritten for the next one, which is the thing this design exists to avoid.
|
|
8
|
+
|
|
9
|
+
So the split is:
|
|
10
|
+
|
|
11
|
+
- **ALK** drives the simulation: who the customer is, when they speak, when it ends.
|
|
12
|
+
- **The world** answers every tool call, through ``handle_tool_call``.
|
|
13
|
+
- **The harness** grades afterwards, from the state the world is left in plus the transcript.
|
|
14
|
+
|
|
15
|
+
What is written here is only the two adapters between the shapes: a scenario becomes an ALK
|
|
16
|
+
``Persona``, and the agent under test becomes an ``AgentWrapper``.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from fi.simulate import Persona, Scenario as AlkScenario
|
|
24
|
+
from fi.simulate.agent.wrapper import AgentInput, AgentResponse, AgentWrapper
|
|
25
|
+
from fi.simulate.environments.chat import ChatEnvironment
|
|
26
|
+
|
|
27
|
+
from ..contract import AgentContract
|
|
28
|
+
from ..scenario import Scenario
|
|
29
|
+
from ..world.runtime import GeneratedWorld
|
|
30
|
+
from .targets import LocalAgent
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def as_persona(scenario: Scenario, simulator_prompt: str = "") -> Persona:
|
|
34
|
+
"""One of our scenarios, in the shape ALK's simulation consumes.
|
|
35
|
+
|
|
36
|
+
``situation`` is the simulator prompt the harness wrote for this agent with the scenario's
|
|
37
|
+
values filled in. ALK wraps it in its own voice-execution rules, so what goes here is only
|
|
38
|
+
what changes per scenario, not a second set of instructions about how to behave on a call.
|
|
39
|
+
|
|
40
|
+
The structured persona is preserved so LiveKit can vary the caller's identity, speech style,
|
|
41
|
+
and scenario metadata instead of flattening every test into the same generic customer.
|
|
42
|
+
"""
|
|
43
|
+
from ..simulator import fill
|
|
44
|
+
|
|
45
|
+
# The LiveKit voice simulator treats ``situation`` as private context rather than a
|
|
46
|
+
# line to recite. Preserve the harness-authored caller rules here: they explicitly keep
|
|
47
|
+
# the simulator in the customer role and stop it from volunteering held-back details.
|
|
48
|
+
# Fall back to the plain scenario instruction for older sessions without a prompt.
|
|
49
|
+
filled = fill(simulator_prompt, scenario.slots())[0] if simulator_prompt else ""
|
|
50
|
+
persona = (
|
|
51
|
+
scenario.persona.model_dump(exclude_none=True)
|
|
52
|
+
if scenario.persona is not None
|
|
53
|
+
else {"name": "customer"}
|
|
54
|
+
)
|
|
55
|
+
return Persona(
|
|
56
|
+
persona=persona,
|
|
57
|
+
situation=filled or scenario.instruction,
|
|
58
|
+
# Said in the person's own terms, because it is read aloud with the situation.
|
|
59
|
+
# ``scenario.tests`` describes what the suite is checking — "agent correctly counts
|
|
60
|
+
# customers filtered by country" — and a person who opens by announcing what the agent
|
|
61
|
+
# is being graded on has told it the answer.
|
|
62
|
+
outcome="get what you came for, or accept that you cannot",
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def as_alk_scenario(
|
|
67
|
+
scenarios: list[Scenario], name: str = "harness", simulator_prompt: str = ""
|
|
68
|
+
) -> AlkScenario:
|
|
69
|
+
return AlkScenario(
|
|
70
|
+
name=name,
|
|
71
|
+
description="generated by the harness",
|
|
72
|
+
dataset=[as_persona(one, simulator_prompt) for one in scenarios],
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _spoken(input: AgentInput) -> str:
|
|
77
|
+
"""What the customer just said, as text.
|
|
78
|
+
|
|
79
|
+
ALK passes a message as a mapping, not a string, and hands the whole history alongside it.
|
|
80
|
+
Passing the mapping straight to a session that expects text fails inside the SDK with a
|
|
81
|
+
redacted TypeError, which says nothing about where it came from.
|
|
82
|
+
"""
|
|
83
|
+
latest = input.new_message
|
|
84
|
+
if isinstance(latest, dict):
|
|
85
|
+
content = latest.get("content")
|
|
86
|
+
if isinstance(content, list):
|
|
87
|
+
content = " ".join(
|
|
88
|
+
part.get("text", "") for part in content if isinstance(part, dict)
|
|
89
|
+
)
|
|
90
|
+
if content:
|
|
91
|
+
return str(content)
|
|
92
|
+
if isinstance(latest, str) and latest:
|
|
93
|
+
return latest
|
|
94
|
+
for message in reversed(input.messages or []):
|
|
95
|
+
if isinstance(message, dict) and message.get("role") != "assistant":
|
|
96
|
+
content = message.get("content")
|
|
97
|
+
if content:
|
|
98
|
+
return str(content)
|
|
99
|
+
return "(the customer said nothing)"
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class ContractAgent(AgentWrapper):
|
|
103
|
+
"""The agent under test, in the shape ALK drives agents by.
|
|
104
|
+
|
|
105
|
+
It holds the same session ``LocalAgent`` uses, so the agent being graded is identical either
|
|
106
|
+
way; what changes is who runs the conversation around it. The tool calls it made are reported
|
|
107
|
+
back to ALK so they appear in the transcript, having already gone through the world.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
def __init__(
|
|
111
|
+
self,
|
|
112
|
+
contract: AgentContract,
|
|
113
|
+
world: GeneratedWorld,
|
|
114
|
+
*,
|
|
115
|
+
model: str | None = None,
|
|
116
|
+
) -> None:
|
|
117
|
+
self.agent = LocalAgent(contract, world, model=model)
|
|
118
|
+
self.world = world
|
|
119
|
+
self._open = False
|
|
120
|
+
|
|
121
|
+
async def call(self, input: AgentInput) -> AgentResponse:
|
|
122
|
+
if not self._open:
|
|
123
|
+
await self.agent.open()
|
|
124
|
+
self._open = True
|
|
125
|
+
|
|
126
|
+
before = len(self.world.calls)
|
|
127
|
+
said = await self.agent.say(_spoken(input))
|
|
128
|
+
made = self.world.calls[before:]
|
|
129
|
+
|
|
130
|
+
return AgentResponse(
|
|
131
|
+
content=said,
|
|
132
|
+
tool_calls=[
|
|
133
|
+
{"name": call.name, "arguments": call.arguments} for call in made
|
|
134
|
+
],
|
|
135
|
+
tool_responses=[
|
|
136
|
+
{
|
|
137
|
+
"name": call.name,
|
|
138
|
+
"content": call.error if not call.ok else str(call.result),
|
|
139
|
+
"success": call.ok,
|
|
140
|
+
}
|
|
141
|
+
for call in made
|
|
142
|
+
],
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
async def aclose(self) -> None:
|
|
146
|
+
if self._open:
|
|
147
|
+
await self.agent.close()
|
|
148
|
+
self._open = False
|
|
149
|
+
|
|
150
|
+
@property
|
|
151
|
+
def spent_usd(self) -> float:
|
|
152
|
+
return self.agent.spent_usd
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
async def simulate(
|
|
156
|
+
scenario: Scenario,
|
|
157
|
+
contract: AgentContract,
|
|
158
|
+
world: GeneratedWorld,
|
|
159
|
+
*,
|
|
160
|
+
model: str | None = None,
|
|
161
|
+
simulator_prompt: str = "",
|
|
162
|
+
) -> tuple[Any, float]:
|
|
163
|
+
"""Run one scenario through ALK's chat simulation against this world.
|
|
164
|
+
|
|
165
|
+
``auto_execute_tools`` is off because the agent's tools are bound to the world already and
|
|
166
|
+
have run by the time it answers. Turning it on would execute every call a second time, which
|
|
167
|
+
for a world that really writes rows means every order placed twice.
|
|
168
|
+
"""
|
|
169
|
+
agent = ContractAgent(contract, world, model=model)
|
|
170
|
+
try:
|
|
171
|
+
report = await ChatEnvironment().run(
|
|
172
|
+
scenario=as_alk_scenario(
|
|
173
|
+
[scenario], name=scenario.name, simulator_prompt=simulator_prompt
|
|
174
|
+
),
|
|
175
|
+
agent_callback=agent,
|
|
176
|
+
environment=world,
|
|
177
|
+
auto_execute_tools=False,
|
|
178
|
+
max_turns=max(2, scenario.max_turns),
|
|
179
|
+
min_turns=2,
|
|
180
|
+
modality=contract.modality or "text",
|
|
181
|
+
)
|
|
182
|
+
finally:
|
|
183
|
+
await agent.aclose()
|
|
184
|
+
return report, agent.spent_usd
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""One scenario, against the real hosted agent, end to end.
|
|
2
|
+
|
|
3
|
+
Everything the harness built is wired together here and then ALK's own voice case places the
|
|
4
|
+
call. The harness does not reimplement any of that: it supplies the world the agent's tools act
|
|
5
|
+
on, the caller's instruction, and the grading afterwards.
|
|
6
|
+
|
|
7
|
+
world + setup ──► webhook ──► public url ──► assistant's own tools repointed
|
|
8
|
+
│
|
|
9
|
+
ALK's voice case places the call ──┘
|
|
10
|
+
│
|
|
11
|
+
the world afterwards + the calls ──► sub-goal checks
|
|
12
|
+
|
|
13
|
+
Run it:
|
|
14
|
+
|
|
15
|
+
set -a; . ./.env.acceptance; set +a
|
|
16
|
+
uv run python -m harness.run.call --name drive_thru --scenario orders_a_big_mac
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import subprocess
|
|
25
|
+
import sys
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
from ..config import artifact_dir
|
|
29
|
+
from ..scenario_tools import load_scenarios
|
|
30
|
+
from .live import grade, wire
|
|
31
|
+
|
|
32
|
+
CASE = os.environ.get("HARNESS_VOICE_CASE", "2.1.2")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# The default runner is part of the harness package and uses ALK's public
|
|
36
|
+
# SimulationSpec/SimulationRunner API. It is overridable only for provider
|
|
37
|
+
# acceptance work; normal customer runs need no second checkout or script.
|
|
38
|
+
VOICE_RUNNER = Path(__file__).with_name("sdk_voice.py")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
LIVE_EVENT = "HARNESS_EXCHANGE "
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def place_the_call(case: str, dry_run: bool = False, on_exchange=None) -> int:
|
|
45
|
+
"""Hand over to ALK's voice case, which owns everything about placing a call."""
|
|
46
|
+
named = os.environ.get("HARNESS_VOICE_RUNNER", "").strip()
|
|
47
|
+
runner = Path(named) if named else VOICE_RUNNER
|
|
48
|
+
if not runner.exists():
|
|
49
|
+
raise RuntimeError(
|
|
50
|
+
f"no voice runner at {runner}. It ships in the harness package; set "
|
|
51
|
+
"HARNESS_VOICE_RUNNER if it lives somewhere else."
|
|
52
|
+
)
|
|
53
|
+
command = [sys.executable, str(runner), case] + (["--dry-run"] if dry_run else [])
|
|
54
|
+
if on_exchange is None:
|
|
55
|
+
return subprocess.call(command)
|
|
56
|
+
process = subprocess.Popen(
|
|
57
|
+
command,
|
|
58
|
+
stdout=subprocess.PIPE,
|
|
59
|
+
stderr=subprocess.STDOUT,
|
|
60
|
+
text=True,
|
|
61
|
+
bufsize=1,
|
|
62
|
+
)
|
|
63
|
+
assert process.stdout is not None
|
|
64
|
+
for line in process.stdout:
|
|
65
|
+
if line.startswith(LIVE_EVENT):
|
|
66
|
+
try:
|
|
67
|
+
on_exchange(json.loads(line[len(LIVE_EVENT) :]))
|
|
68
|
+
except (json.JSONDecodeError, TypeError):
|
|
69
|
+
pass
|
|
70
|
+
else:
|
|
71
|
+
print(line, end="", flush=True)
|
|
72
|
+
return process.wait()
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def main(argv: list[str] | None = None) -> int:
|
|
76
|
+
parser = argparse.ArgumentParser(prog="agent-harness-call", description=__doc__)
|
|
77
|
+
parser.add_argument("--name", required=True, help="which agent")
|
|
78
|
+
parser.add_argument("--scenario", required=True, help="which scenario, by name")
|
|
79
|
+
parser.add_argument("--case", default=CASE, help="ALK voice case id")
|
|
80
|
+
parser.add_argument(
|
|
81
|
+
"--dry-run",
|
|
82
|
+
action="store_true",
|
|
83
|
+
help="wire everything up but do not place the call",
|
|
84
|
+
)
|
|
85
|
+
args = parser.parse_args(argv)
|
|
86
|
+
|
|
87
|
+
root = artifact_dir(args.name)
|
|
88
|
+
written = load_scenarios(root)
|
|
89
|
+
scenario = next((one for one in written if one.name == args.scenario), None)
|
|
90
|
+
if scenario is None:
|
|
91
|
+
print(
|
|
92
|
+
f"no scenario called {args.scenario!r}. There is: "
|
|
93
|
+
+ ", ".join(one.name for one in written),
|
|
94
|
+
file=sys.stderr,
|
|
95
|
+
)
|
|
96
|
+
return 1
|
|
97
|
+
|
|
98
|
+
world, instruction, webhook, tunnel, url, moved = wire(scenario, root)
|
|
99
|
+
try:
|
|
100
|
+
print(f"agent: {args.name}")
|
|
101
|
+
print(f"scenario: {scenario.name}")
|
|
102
|
+
print(f"webhook: {url}/tool")
|
|
103
|
+
print(f"repointed: {', '.join(moved)}")
|
|
104
|
+
print(f"sub-goals: {', '.join(scenario.sub_goals)}\n")
|
|
105
|
+
|
|
106
|
+
# The caller's instruction reaches the voice case through the environment, so nothing
|
|
107
|
+
# about how a simulated caller behaves is decided twice.
|
|
108
|
+
os.environ["HARNESS_INSTRUCTION"] = instruction
|
|
109
|
+
os.environ["HARNESS_SCENARIO"] = scenario.name
|
|
110
|
+
# The caller prompt is a template the harness fills, never prose it composes, so
|
|
111
|
+
# the generated template travels to the call with everything else.
|
|
112
|
+
# The caller is never handed the grader's pass question. `tests` is written about
|
|
113
|
+
# the agent in the third person, so as an objective it reads as a rubric rather
|
|
114
|
+
# than a motive. What this person wants is already in the instruction.
|
|
115
|
+
os.environ.pop("HARNESS_OUTCOME", None)
|
|
116
|
+
os.environ["HARNESS_PERSONA"] = json.dumps(
|
|
117
|
+
scenario.persona.model_dump(exclude_none=True)
|
|
118
|
+
if scenario.persona is not None
|
|
119
|
+
else {"name": "customer"}
|
|
120
|
+
)
|
|
121
|
+
os.environ["HARNESS_INITIAL_MESSAGE"] = (
|
|
122
|
+
scenario.persona.initial_message if scenario.persona is not None else ""
|
|
123
|
+
)
|
|
124
|
+
os.environ["HARNESS_FIXTURE"] = json.dumps(
|
|
125
|
+
scenario.fixture, ensure_ascii=False, default=str
|
|
126
|
+
)
|
|
127
|
+
os.environ["HARNESS_SCRIPTED_CALLER"] = json.dumps(
|
|
128
|
+
scenario.persona.scripted_caller
|
|
129
|
+
if scenario.persona is not None
|
|
130
|
+
and scenario.persona.scripted_caller is not None
|
|
131
|
+
else {}
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
code = place_the_call(args.case, dry_run=args.dry_run)
|
|
135
|
+
if args.dry_run:
|
|
136
|
+
print("\ndry run: nothing was called, and the world is untouched.")
|
|
137
|
+
return code
|
|
138
|
+
|
|
139
|
+
result = grade(scenario, world, root)
|
|
140
|
+
print()
|
|
141
|
+
print(result.line())
|
|
142
|
+
for one in result.settled:
|
|
143
|
+
print(one.line())
|
|
144
|
+
for name in result.judged:
|
|
145
|
+
print(f" [?] {name} — judged, not graded here")
|
|
146
|
+
print("\nwhat the agent actually did:")
|
|
147
|
+
for call in result.calls or ["(no tool calls reached the world)"]:
|
|
148
|
+
print(f" {call}")
|
|
149
|
+
return 0 if result.settled and result.met == len(result.settled) else 2
|
|
150
|
+
finally:
|
|
151
|
+
webhook.stop()
|
|
152
|
+
if tunnel is not None:
|
|
153
|
+
tunnel.terminate()
|
|
154
|
+
if (root / "environment.json").exists():
|
|
155
|
+
from ..provision import stop_runtime
|
|
156
|
+
|
|
157
|
+
stop_runtime(root)
|
|
158
|
+
world.close()
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
if __name__ == "__main__":
|
|
162
|
+
raise SystemExit(main())
|