agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,340 @@
|
|
|
1
|
+
"""Serving a real voice agent's tool calls from a generated world.
|
|
2
|
+
|
|
3
|
+
A hosted voice agent executes its tools by calling a webhook. So the whole integration is one
|
|
4
|
+
thing: stand up that webhook, and answer it from the world instead of from canned responses.
|
|
5
|
+
|
|
6
|
+
That single swap is what the environment was built for. The previous run's known issues were all
|
|
7
|
+
the same defect wearing different clothes:
|
|
8
|
+
|
|
9
|
+
- *"Mocked tools always succeed, including removing an item that was never added."*
|
|
10
|
+
- *"Mock responses do not vary by argument, so read-after-write flows are wrong."*
|
|
11
|
+
- *"World state does not change unless a scenario sets state_updates, which is often empty."*
|
|
12
|
+
|
|
13
|
+
A world that really holds rows and can really refuse answers all three, because the reply the
|
|
14
|
+
agent hears is produced by running the call rather than by looking it up.
|
|
15
|
+
|
|
16
|
+
Nothing here decides pass or fail. Grading reads the world afterwards and the calls this server
|
|
17
|
+
recorded, through the same sub-goal checks every other run uses.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
import logging
|
|
24
|
+
import os
|
|
25
|
+
import threading
|
|
26
|
+
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
27
|
+
from typing import Any, Mapping
|
|
28
|
+
|
|
29
|
+
from ..world.runtime import GeneratedWorld
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
VAPI_API = os.environ.get("VAPI_API_BASE_URL", "https://api.vapi.ai").rstrip("/")
|
|
34
|
+
|
|
35
|
+
# Vapi's edge rejects the default urllib User-Agent with a 403 that says nothing about why, while
|
|
36
|
+
# the identical request from curl succeeds. Sending one is the whole fix.
|
|
37
|
+
_AGENT = "alk-harness/0.1"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class WorldWebhook:
|
|
41
|
+
"""The webhook a hosted agent calls, answered by a generated world.
|
|
42
|
+
|
|
43
|
+
One world at a time. ``bind`` swaps which world is live between scenarios, so the assistant
|
|
44
|
+
stays configured while every scenario still starts from its own restored copy.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(self, host: str | None = None, port: int | None = None) -> None:
|
|
48
|
+
# A container must bind 0.0.0.0 on a fixed port or nothing outside can
|
|
49
|
+
# be configured to call it; a laptop keeps loopback and an ephemeral port.
|
|
50
|
+
host = (
|
|
51
|
+
host
|
|
52
|
+
if host is not None
|
|
53
|
+
else os.environ.get("HARNESS_WEBHOOK_HOST", "127.0.0.1")
|
|
54
|
+
)
|
|
55
|
+
port = (
|
|
56
|
+
port
|
|
57
|
+
if port is not None
|
|
58
|
+
else int(os.environ.get("HARNESS_WEBHOOK_PORT", "0"))
|
|
59
|
+
)
|
|
60
|
+
self._world: GeneratedWorld | None = None
|
|
61
|
+
self._lock = threading.Lock()
|
|
62
|
+
try:
|
|
63
|
+
server = HTTPServer((host, port), _handler_for(self))
|
|
64
|
+
except OSError:
|
|
65
|
+
# A leftover server from an earlier run must not block this one; any free port works
|
|
66
|
+
# because the public URL is discovered after binding.
|
|
67
|
+
logger.warning("port %s busy, binding an ephemeral port instead", port)
|
|
68
|
+
server = HTTPServer((host, 0), _handler_for(self))
|
|
69
|
+
self._server = server
|
|
70
|
+
self.port = server.server_address[1]
|
|
71
|
+
self._thread = threading.Thread(target=server.serve_forever, daemon=True)
|
|
72
|
+
|
|
73
|
+
def start(self) -> "WorldWebhook":
|
|
74
|
+
self._thread.start()
|
|
75
|
+
logger.info("world webhook listening on port %s", self.port)
|
|
76
|
+
return self
|
|
77
|
+
|
|
78
|
+
def stop(self) -> None:
|
|
79
|
+
self._server.shutdown()
|
|
80
|
+
self._server.server_close()
|
|
81
|
+
|
|
82
|
+
def bind(self, world: GeneratedWorld) -> None:
|
|
83
|
+
"""Make one world live. Its own call log is what grading reads afterwards."""
|
|
84
|
+
with self._lock:
|
|
85
|
+
self._world = world
|
|
86
|
+
world.reset()
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def calls(self) -> list[Any]:
|
|
90
|
+
with self._lock:
|
|
91
|
+
return list(self._world.calls) if self._world else []
|
|
92
|
+
|
|
93
|
+
def respond(
|
|
94
|
+
self,
|
|
95
|
+
name: str,
|
|
96
|
+
arguments: Mapping[str, Any],
|
|
97
|
+
*,
|
|
98
|
+
session_id: str = "harness",
|
|
99
|
+
) -> str:
|
|
100
|
+
"""Answer one tool call by running it.
|
|
101
|
+
|
|
102
|
+
A refusal is returned as the answer, not as an error: the agent has to hear "that item is
|
|
103
|
+
unavailable" and cope with it, which is the whole reason the world can say no. What it
|
|
104
|
+
must never hear is an acknowledgement for something that did not happen.
|
|
105
|
+
"""
|
|
106
|
+
with self._lock:
|
|
107
|
+
world = self._world
|
|
108
|
+
if world is None:
|
|
109
|
+
return "the environment is not ready"
|
|
110
|
+
|
|
111
|
+
# Caller hydration is an adapter concern, not a contract tool. The local
|
|
112
|
+
# LiveKit worker normally gets this from its demo API before exposing any
|
|
113
|
+
# conversational tools. Resolve the same safe profile from the generated
|
|
114
|
+
# world without adding a call that scenario grading would mistake for an
|
|
115
|
+
# agent action.
|
|
116
|
+
if name == "lookup_rider_by_phone":
|
|
117
|
+
phone = str(arguments.get("phone") or "")
|
|
118
|
+
state = world.observe().state
|
|
119
|
+
user = next(
|
|
120
|
+
(
|
|
121
|
+
row
|
|
122
|
+
for row in state.get("users", [])
|
|
123
|
+
if str(row.get("phone")) == phone
|
|
124
|
+
),
|
|
125
|
+
None,
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
def active_booking_ref() -> Any:
|
|
129
|
+
if user is None:
|
|
130
|
+
return None
|
|
131
|
+
active = [
|
|
132
|
+
row
|
|
133
|
+
for row in state.get("bookings", [])
|
|
134
|
+
if row.get("rider_id") == user.get("rider_id")
|
|
135
|
+
and str(row.get("status") or "").lower()
|
|
136
|
+
not in {"cancelled", "canceled", "completed"}
|
|
137
|
+
]
|
|
138
|
+
active.sort(
|
|
139
|
+
key=lambda row: str(row.get("created_at") or ""), reverse=True
|
|
140
|
+
)
|
|
141
|
+
return active[0].get("booking_ref") if active else None
|
|
142
|
+
|
|
143
|
+
forward = getattr(world, "forward", None)
|
|
144
|
+
if callable(forward):
|
|
145
|
+
hydrated = forward(
|
|
146
|
+
name,
|
|
147
|
+
arguments,
|
|
148
|
+
record=False,
|
|
149
|
+
session_id=session_id,
|
|
150
|
+
)
|
|
151
|
+
if hydrated.ok:
|
|
152
|
+
result = hydrated.result
|
|
153
|
+
if isinstance(result, dict):
|
|
154
|
+
result = {**result, "booking_ref": active_booking_ref()}
|
|
155
|
+
return (
|
|
156
|
+
result
|
|
157
|
+
if isinstance(result, str)
|
|
158
|
+
else json.dumps(result, default=str)
|
|
159
|
+
)
|
|
160
|
+
return hydrated.error
|
|
161
|
+
if user is None:
|
|
162
|
+
return json.dumps({"rider_id": None, "phone": phone})
|
|
163
|
+
market = next(
|
|
164
|
+
(
|
|
165
|
+
row
|
|
166
|
+
for row in state.get("market_config", [])
|
|
167
|
+
if row.get("market") == user.get("default_market")
|
|
168
|
+
),
|
|
169
|
+
{},
|
|
170
|
+
)
|
|
171
|
+
# A caller asking about or cancelling an existing ride does not know an internal
|
|
172
|
+
# booking reference. Real agent backends hydrate the active trip alongside ANI
|
|
173
|
+
# identity; expose the same seeded relationship from the generated world so the
|
|
174
|
+
# shipped agent can invoke its own reference-based tools without fixture leakage in
|
|
175
|
+
# the conversation.
|
|
176
|
+
return json.dumps(
|
|
177
|
+
{
|
|
178
|
+
**user,
|
|
179
|
+
"cash_supported_in_market": bool(market.get("cash_supported")),
|
|
180
|
+
"accessibility_needs": [],
|
|
181
|
+
"booking_ref": active_booking_ref(),
|
|
182
|
+
},
|
|
183
|
+
default=str,
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
# ``world.call`` distinguishes real dependency endpoints from actions executed inside
|
|
187
|
+
# the submitted worker and mirrored here only for evidence. Bypassing it through
|
|
188
|
+
# ``forward`` made a valid local action look like a missing HTTP endpoint and caused the
|
|
189
|
+
# agent to retry or abort after it had already updated its own state.
|
|
190
|
+
done = world.handle_tool_call({"name": name, "arguments": dict(arguments)})
|
|
191
|
+
if done is None:
|
|
192
|
+
return f"there is no tool called {name}"
|
|
193
|
+
return done.content or ("done" if done.success else "that could not be done")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _handler_for(owner: "WorldWebhook"):
|
|
197
|
+
class Handler(BaseHTTPRequestHandler):
|
|
198
|
+
def log_message(self, *args: Any) -> None: # silence per-request stderr noise
|
|
199
|
+
return
|
|
200
|
+
|
|
201
|
+
def do_POST(self) -> None: # noqa: N802 - required name
|
|
202
|
+
length = int(self.headers.get("Content-Length") or 0)
|
|
203
|
+
raw = self.rfile.read(length) if length else b"{}"
|
|
204
|
+
try:
|
|
205
|
+
payload = json.loads(raw or b"{}")
|
|
206
|
+
except json.JSONDecodeError:
|
|
207
|
+
payload = {}
|
|
208
|
+
|
|
209
|
+
calls = tool_calls(payload)
|
|
210
|
+
session_id = self.headers.get("x-session-id") or "harness"
|
|
211
|
+
if not calls:
|
|
212
|
+
# An agent whose tools are its own HTTP API asks differently: the tool is the
|
|
213
|
+
# path and the body is the arguments, with the answer expected back plainly.
|
|
214
|
+
# Serving both shapes is what lets a world stand in for such an API without the
|
|
215
|
+
# agent being changed to suit us.
|
|
216
|
+
name = self.path.strip("/").split("?")[0]
|
|
217
|
+
if name:
|
|
218
|
+
answer = owner.respond(
|
|
219
|
+
name,
|
|
220
|
+
payload if isinstance(payload, dict) else {},
|
|
221
|
+
session_id=session_id,
|
|
222
|
+
)
|
|
223
|
+
plain = json.dumps(_as_body(answer)).encode()
|
|
224
|
+
self.send_response(200)
|
|
225
|
+
self.send_header("Content-Type", "application/json")
|
|
226
|
+
self.send_header("Content-Length", str(len(plain)))
|
|
227
|
+
self.end_headers()
|
|
228
|
+
self.wfile.write(plain)
|
|
229
|
+
return
|
|
230
|
+
|
|
231
|
+
results = [
|
|
232
|
+
{
|
|
233
|
+
"toolCallId": call_id,
|
|
234
|
+
"result": owner.respond(name, arguments, session_id=session_id),
|
|
235
|
+
}
|
|
236
|
+
for call_id, name, arguments in calls
|
|
237
|
+
]
|
|
238
|
+
body = json.dumps({"results": results}).encode()
|
|
239
|
+
self.send_response(200)
|
|
240
|
+
self.send_header("Content-Type", "application/json")
|
|
241
|
+
self.send_header("Content-Length", str(len(body)))
|
|
242
|
+
self.end_headers()
|
|
243
|
+
self.wfile.write(body)
|
|
244
|
+
|
|
245
|
+
return Handler
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _as_body(answer: str) -> Any:
|
|
249
|
+
"""A tool's answer as a JSON body, keeping structure when the handler produced any.
|
|
250
|
+
|
|
251
|
+
Handlers return text because that is what a spoken agent hears. An HTTP tool API expects an
|
|
252
|
+
object, so a JSON answer is passed through as itself and anything else is wrapped, rather
|
|
253
|
+
than a caller having to parse a string out of a string.
|
|
254
|
+
"""
|
|
255
|
+
try:
|
|
256
|
+
parsed = json.loads(answer)
|
|
257
|
+
except (json.JSONDecodeError, TypeError):
|
|
258
|
+
return {"result": answer}
|
|
259
|
+
return parsed if isinstance(parsed, (dict, list)) else {"result": parsed}
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def tool_calls(payload: Mapping[str, Any]) -> list[tuple[str, str, dict[str, Any]]]:
|
|
263
|
+
"""Pull (id, name, arguments) out of a provider's tool-call webhook body."""
|
|
264
|
+
message = payload.get("message") or payload
|
|
265
|
+
raw = message.get("toolCalls") or message.get("toolCallList") or []
|
|
266
|
+
found: list[tuple[str, str, dict[str, Any]]] = []
|
|
267
|
+
for entry in raw if isinstance(raw, list) else []:
|
|
268
|
+
if not isinstance(entry, Mapping):
|
|
269
|
+
continue
|
|
270
|
+
function = entry.get("function") or {}
|
|
271
|
+
name = str(function.get("name") or entry.get("name") or "")
|
|
272
|
+
arguments = function.get("arguments") or entry.get("arguments") or {}
|
|
273
|
+
if isinstance(arguments, str):
|
|
274
|
+
try:
|
|
275
|
+
arguments = json.loads(arguments)
|
|
276
|
+
except json.JSONDecodeError:
|
|
277
|
+
arguments = {"_raw": arguments}
|
|
278
|
+
if name:
|
|
279
|
+
found.append((str(entry.get("id") or ""), name, dict(arguments)))
|
|
280
|
+
return found
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def pointed_at(tools: list[dict[str, Any]], webhook_url: str) -> list[dict[str, Any]]:
|
|
284
|
+
"""The agent's own tools, with only where they are answered changed.
|
|
285
|
+
|
|
286
|
+
The assistant under test already has its tools — the names, the arguments, the enums are the
|
|
287
|
+
agent's, defined by whoever built it. Redefining them here would mean testing an agent we
|
|
288
|
+
wrote rather than theirs, and any drift between the two would show up as a finding about
|
|
289
|
+
them. So nothing is rebuilt: the one thing that changes is the address the call goes to.
|
|
290
|
+
"""
|
|
291
|
+
repointed: list[dict[str, Any]] = []
|
|
292
|
+
for tool in tools:
|
|
293
|
+
moved = json.loads(json.dumps(tool))
|
|
294
|
+
moved.setdefault("server", {})["url"] = f"{webhook_url.rstrip('/')}/tool"
|
|
295
|
+
repointed.append(moved)
|
|
296
|
+
return repointed
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def fetch_assistant(assistant_id: str, api_key: str) -> dict[str, Any]:
|
|
300
|
+
"""The assistant as it stands, so its own tools can be read rather than guessed."""
|
|
301
|
+
import urllib.request
|
|
302
|
+
|
|
303
|
+
request = urllib.request.Request(
|
|
304
|
+
f"{VAPI_API}/assistant/{assistant_id}",
|
|
305
|
+
headers={"Authorization": f"Bearer {api_key}", "User-Agent": _AGENT},
|
|
306
|
+
)
|
|
307
|
+
with urllib.request.urlopen(request, timeout=20) as answer:
|
|
308
|
+
return json.loads(answer.read())
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def repoint_assistant(assistant_id: str, api_key: str, webhook_url: str) -> list[str]:
|
|
312
|
+
"""Send the assistant's existing tool calls to our webhook. Returns the tools moved."""
|
|
313
|
+
import urllib.request
|
|
314
|
+
|
|
315
|
+
assistant = fetch_assistant(assistant_id, api_key)
|
|
316
|
+
tools = (assistant.get("model") or {}).get("tools") or []
|
|
317
|
+
if not tools:
|
|
318
|
+
raise RuntimeError(
|
|
319
|
+
f"assistant {assistant_id} has no tools, so there is nothing for the environment "
|
|
320
|
+
"to answer. It is the agent's own tools that get repointed, not ones we add."
|
|
321
|
+
)
|
|
322
|
+
model = json.loads(json.dumps(assistant.get("model") or {}))
|
|
323
|
+
model["tools"] = pointed_at(tools, webhook_url)
|
|
324
|
+
|
|
325
|
+
body = json.dumps({"model": model}).encode()
|
|
326
|
+
request = urllib.request.Request(
|
|
327
|
+
f"{VAPI_API}/assistant/{assistant_id}",
|
|
328
|
+
data=body,
|
|
329
|
+
method="PATCH",
|
|
330
|
+
headers={
|
|
331
|
+
"Authorization": f"Bearer {api_key}",
|
|
332
|
+
"Content-Type": "application/json",
|
|
333
|
+
"User-Agent": _AGENT,
|
|
334
|
+
},
|
|
335
|
+
)
|
|
336
|
+
with urllib.request.urlopen(request, timeout=20) as answer:
|
|
337
|
+
answer.read()
|
|
338
|
+
return [
|
|
339
|
+
str((one.get("function") or {}).get("name") or "") for one in model["tools"]
|
|
340
|
+
]
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Runtime-provider boundary for ALK-owned test environments.
|
|
2
|
+
|
|
3
|
+
Providers decide *where* a sealed environment runs. They do not decide how an agent is
|
|
4
|
+
understood, how scenarios are written, or how results are graded. The local provider below
|
|
5
|
+
adapts the proven repository/Compose provisioner; the hosted sandbox implements the same port.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any, Protocol
|
|
14
|
+
|
|
15
|
+
from pydantic import BaseModel, Field, JsonValue
|
|
16
|
+
|
|
17
|
+
from .bundle import EnvironmentBundle
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class RuntimeState(str, Enum):
|
|
21
|
+
PREPARING = "preparing"
|
|
22
|
+
READY = "ready"
|
|
23
|
+
UNHEALTHY = "unhealthy"
|
|
24
|
+
STOPPED = "stopped"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class RuntimeEndpoint(BaseModel):
|
|
28
|
+
capability: str
|
|
29
|
+
protocol: str
|
|
30
|
+
address: str
|
|
31
|
+
configuration_name: str | None = None
|
|
32
|
+
metadata: dict[str, JsonValue] = Field(default_factory=dict)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class EnvironmentRuntime(BaseModel):
|
|
36
|
+
runtime_id: str
|
|
37
|
+
provider: str
|
|
38
|
+
bundle_digest: str
|
|
39
|
+
state: RuntimeState
|
|
40
|
+
endpoints: dict[str, RuntimeEndpoint] = Field(default_factory=dict)
|
|
41
|
+
metadata: dict[str, JsonValue] = Field(default_factory=dict)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class RuntimeProvider(Protocol):
|
|
45
|
+
"""Execution-location port implemented locally and by the hosted sandbox fleet."""
|
|
46
|
+
|
|
47
|
+
name: str
|
|
48
|
+
|
|
49
|
+
async def provision(
|
|
50
|
+
self,
|
|
51
|
+
bundle: EnvironmentBundle,
|
|
52
|
+
*,
|
|
53
|
+
source: Path,
|
|
54
|
+
work_directory: Path,
|
|
55
|
+
contract: Any | None = None,
|
|
56
|
+
) -> EnvironmentRuntime: ...
|
|
57
|
+
|
|
58
|
+
async def reset(
|
|
59
|
+
self, runtime: EnvironmentRuntime, *, work_directory: Path
|
|
60
|
+
) -> None: ...
|
|
61
|
+
|
|
62
|
+
async def healthy(
|
|
63
|
+
self, runtime: EnvironmentRuntime, *, work_directory: Path
|
|
64
|
+
) -> bool: ...
|
|
65
|
+
|
|
66
|
+
async def close(
|
|
67
|
+
self, runtime: EnvironmentRuntime, *, work_directory: Path
|
|
68
|
+
) -> None: ...
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class LocalComposeRuntimeProvider:
|
|
72
|
+
"""Run one repository environment as an isolated Docker Compose project.
|
|
73
|
+
|
|
74
|
+
The adapter delegates lifecycle mechanics to ``harness.provision``, which gives each run a
|
|
75
|
+
unique project, allocates ports, waits for declared health checks, fingerprints source reuse,
|
|
76
|
+
and removes volumes during cleanup. Only endpoint names and addresses cross this boundary;
|
|
77
|
+
resolved credentials remain process-local.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
name = "local-compose"
|
|
81
|
+
|
|
82
|
+
async def provision(
|
|
83
|
+
self,
|
|
84
|
+
bundle: EnvironmentBundle,
|
|
85
|
+
*,
|
|
86
|
+
source: Path,
|
|
87
|
+
work_directory: Path,
|
|
88
|
+
contract: Any | None = None,
|
|
89
|
+
) -> EnvironmentRuntime:
|
|
90
|
+
from .provision import provision
|
|
91
|
+
|
|
92
|
+
environment = await asyncio.to_thread(
|
|
93
|
+
provision, source, work_directory, contract
|
|
94
|
+
)
|
|
95
|
+
endpoints: dict[str, RuntimeEndpoint] = {}
|
|
96
|
+
overrides = dict(environment.overrides)
|
|
97
|
+
for capability, definition in bundle.capabilities.items():
|
|
98
|
+
address = ""
|
|
99
|
+
if definition.configuration_name:
|
|
100
|
+
address = overrides.get(definition.configuration_name, "")
|
|
101
|
+
if not address and len(overrides) == 1:
|
|
102
|
+
address = next(iter(overrides.values()))
|
|
103
|
+
discovered = next(
|
|
104
|
+
(
|
|
105
|
+
endpoint
|
|
106
|
+
for endpoint in environment.service_endpoints
|
|
107
|
+
if endpoint["service"] == definition.service
|
|
108
|
+
and endpoint["container_port"] == definition.container_port
|
|
109
|
+
),
|
|
110
|
+
None,
|
|
111
|
+
)
|
|
112
|
+
if not address and discovered:
|
|
113
|
+
address = str(discovered["external_address"])
|
|
114
|
+
if address:
|
|
115
|
+
endpoints[capability] = RuntimeEndpoint(
|
|
116
|
+
capability=capability,
|
|
117
|
+
protocol=definition.protocol.value,
|
|
118
|
+
address=address,
|
|
119
|
+
configuration_name=definition.configuration_name,
|
|
120
|
+
metadata=(
|
|
121
|
+
{
|
|
122
|
+
"service": str(discovered["service"]),
|
|
123
|
+
"kind": str(discovered["kind"]),
|
|
124
|
+
}
|
|
125
|
+
if discovered
|
|
126
|
+
else {}
|
|
127
|
+
),
|
|
128
|
+
)
|
|
129
|
+
return EnvironmentRuntime(
|
|
130
|
+
runtime_id=environment.project,
|
|
131
|
+
provider=self.name,
|
|
132
|
+
bundle_digest=bundle.digest,
|
|
133
|
+
state=RuntimeState.READY,
|
|
134
|
+
endpoints=endpoints,
|
|
135
|
+
metadata={
|
|
136
|
+
"services": environment.services,
|
|
137
|
+
"provision_seconds": environment.provision_seconds,
|
|
138
|
+
"managed": environment.managed,
|
|
139
|
+
},
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
async def reset(self, runtime: EnvironmentRuntime, *, work_directory: Path) -> None:
|
|
143
|
+
from .provision import reset
|
|
144
|
+
|
|
145
|
+
environment = await asyncio.to_thread(reset, work_directory)
|
|
146
|
+
runtime.state = (
|
|
147
|
+
RuntimeState.READY if environment.running else RuntimeState.UNHEALTHY
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
async def healthy(
|
|
151
|
+
self, runtime: EnvironmentRuntime, *, work_directory: Path
|
|
152
|
+
) -> bool:
|
|
153
|
+
from .provision import healthy
|
|
154
|
+
|
|
155
|
+
is_healthy = await asyncio.to_thread(healthy, work_directory)
|
|
156
|
+
runtime.state = RuntimeState.READY if is_healthy else RuntimeState.UNHEALTHY
|
|
157
|
+
return is_healthy
|
|
158
|
+
|
|
159
|
+
async def close(self, runtime: EnvironmentRuntime, *, work_directory: Path) -> None:
|
|
160
|
+
from .provision import stop
|
|
161
|
+
|
|
162
|
+
await asyncio.to_thread(stop, work_directory)
|
|
163
|
+
runtime.state = RuntimeState.STOPPED
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
__all__ = [
|
|
167
|
+
"EnvironmentRuntime",
|
|
168
|
+
"LocalComposeRuntimeProvider",
|
|
169
|
+
"RuntimeEndpoint",
|
|
170
|
+
"RuntimeProvider",
|
|
171
|
+
"RuntimeState",
|
|
172
|
+
]
|