agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
"""One scenario, against the real hosted agent, in the environment the harness built.
|
|
2
|
+
|
|
3
|
+
The harness wires the whole thing rather than leaving it to be assembled by hand:
|
|
4
|
+
|
|
5
|
+
1. restore the world and apply the scenario's setup
|
|
6
|
+
2. stand the webhook up and bind that world to it
|
|
7
|
+
3. expose it publicly, because a hosted agent has to reach it
|
|
8
|
+
4. point the assistant's **own** tools at that address — nothing about the agent is redefined
|
|
9
|
+
5. run ALK's voice case with the scenario's instruction driving the simulated caller
|
|
10
|
+
6. grade from the world afterwards and the calls the webhook recorded
|
|
11
|
+
|
|
12
|
+
Steps 1, 2, 4 and 6 are the whole difference from what existed before: the agent's tool calls now
|
|
13
|
+
land in a database that can refuse, instead of in canned responses that always succeed.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import os
|
|
19
|
+
import re
|
|
20
|
+
import shutil
|
|
21
|
+
import subprocess
|
|
22
|
+
import time
|
|
23
|
+
import uuid
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from ..simulator_voice import fixture_caller_phone
|
|
28
|
+
from ..catalogue import load_catalogue
|
|
29
|
+
from ..checks import Outcome, run_check
|
|
30
|
+
from ..folder import apply_setup, check_ready
|
|
31
|
+
from ..scenario import Scenario
|
|
32
|
+
from ..simulator import fill, load_simulator_prompt
|
|
33
|
+
from ..world.runtime import GeneratedWorld
|
|
34
|
+
from ..world.snapshot import restore
|
|
35
|
+
from .voice import WorldWebhook, repoint_assistant
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class LiveRun:
|
|
40
|
+
"""What a live call left behind."""
|
|
41
|
+
|
|
42
|
+
scenario: str
|
|
43
|
+
settled: list[Outcome] = field(default_factory=list)
|
|
44
|
+
judged: list[str] = field(default_factory=list)
|
|
45
|
+
calls: list[str] = field(default_factory=list)
|
|
46
|
+
ended: str = ""
|
|
47
|
+
problems: list[str] = field(default_factory=list)
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def met(self) -> int:
|
|
51
|
+
return sum(1 for one in self.settled if one.held)
|
|
52
|
+
|
|
53
|
+
def line(self) -> str:
|
|
54
|
+
mark = "PASS" if self.settled and self.met == len(self.settled) else "FAIL"
|
|
55
|
+
if self.problems:
|
|
56
|
+
mark = "VOID"
|
|
57
|
+
return f"{mark} {self.scenario} {self.met}/{len(self.settled)} sub-goals settled by code"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def scoped_agent_name(base: str, scenario: str) -> str:
|
|
61
|
+
"""A dispatch name owned by one worker lifetime, never a stale registration."""
|
|
62
|
+
clean_base = re.sub(r"[^a-zA-Z0-9_-]+", "-", base).strip("-") or "agent"
|
|
63
|
+
clean_case = re.sub(r"[^a-zA-Z0-9_-]+", "-", scenario).strip("-") or "case"
|
|
64
|
+
return f"{clean_base[:28]}-{clean_case[:18]}-{uuid.uuid4().hex[:8]}"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def public_url(
|
|
68
|
+
port: int, *, wait: float = 30.0, tries: int = 3
|
|
69
|
+
) -> tuple[str, subprocess.Popen | None]:
|
|
70
|
+
"""A publicly reachable address for the webhook, and the process holding it open.
|
|
71
|
+
|
|
72
|
+
A hosted agent runs on somebody else's infrastructure, so a loopback address is unreachable
|
|
73
|
+
to it. ``cloudflared`` is what the previous runs used; anything giving a public URL works, and
|
|
74
|
+
``HARNESS_WEBHOOK_URL`` skips this entirely when a tunnel is already running.
|
|
75
|
+
|
|
76
|
+
Retried, because a free tunnel is the least reliable thing in the whole path and it fails
|
|
77
|
+
before anything interesting has happened. One slow handshake should not read as a scenario
|
|
78
|
+
the agent failed, and on a suite of forty it would not fail once.
|
|
79
|
+
"""
|
|
80
|
+
named = os.environ.get("HARNESS_WEBHOOK_URL", "").strip()
|
|
81
|
+
if named:
|
|
82
|
+
return named, None
|
|
83
|
+
if not shutil.which("cloudflared"):
|
|
84
|
+
raise RuntimeError(
|
|
85
|
+
"no way to expose the webhook publicly. Either install cloudflared "
|
|
86
|
+
"(brew install cloudflared) or set HARNESS_WEBHOOK_URL to a tunnel you already have."
|
|
87
|
+
)
|
|
88
|
+
for attempt in range(max(1, tries)):
|
|
89
|
+
found, process = _tunnel(port, wait)
|
|
90
|
+
if found:
|
|
91
|
+
return found, process
|
|
92
|
+
if process is not None:
|
|
93
|
+
process.terminate()
|
|
94
|
+
if attempt + 1 < tries:
|
|
95
|
+
time.sleep(2.0)
|
|
96
|
+
raise RuntimeError(
|
|
97
|
+
f"cloudflared did not report a public URL in {tries} attempts. The tunnel is the "
|
|
98
|
+
"flakiest part of this path; set HARNESS_WEBHOOK_URL to one you control to skip it."
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _tunnel(port: int, wait: float) -> tuple[str, subprocess.Popen | None]:
|
|
103
|
+
"""One attempt at a tunnel: the URL if it came up, and the process either way."""
|
|
104
|
+
process = subprocess.Popen(
|
|
105
|
+
["cloudflared", "tunnel", "--url", f"http://127.0.0.1:{port}"],
|
|
106
|
+
stdout=subprocess.PIPE,
|
|
107
|
+
stderr=subprocess.STDOUT,
|
|
108
|
+
text=True,
|
|
109
|
+
)
|
|
110
|
+
deadline = time.time() + wait
|
|
111
|
+
while time.time() < deadline:
|
|
112
|
+
line = process.stdout.readline() if process.stdout else ""
|
|
113
|
+
if not line and process.poll() is not None:
|
|
114
|
+
return "", process
|
|
115
|
+
if "trycloudflare.com" in line:
|
|
116
|
+
for word in line.split():
|
|
117
|
+
if word.startswith("https://") and "trycloudflare.com" in word:
|
|
118
|
+
return word.strip(), process
|
|
119
|
+
return "", process
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def prepare(scenario: Scenario, world_root: Path) -> tuple[GeneratedWorld, str]:
|
|
123
|
+
"""The world this scenario runs in, and what the simulated caller is told.
|
|
124
|
+
|
|
125
|
+
The instruction is the scenario's values filled into the simulator prompt the environment
|
|
126
|
+
step wrote. Nothing about how a caller behaves is decided here; that belongs to the prompt.
|
|
127
|
+
"""
|
|
128
|
+
world = restore(world_root)
|
|
129
|
+
world.reset()
|
|
130
|
+
applied = apply_setup(scenario, world)
|
|
131
|
+
if not applied.ok:
|
|
132
|
+
raise RuntimeError(f"the scenario's setup did not run: {applied.said}")
|
|
133
|
+
ready = check_ready(scenario, world)
|
|
134
|
+
if not ready.ok:
|
|
135
|
+
raise RuntimeError(
|
|
136
|
+
f"the world is not ready for this scenario: {ready.said}. Running it would test us "
|
|
137
|
+
"rather than the agent."
|
|
138
|
+
)
|
|
139
|
+
# The setup's own calls are not the agent's.
|
|
140
|
+
world.calls = []
|
|
141
|
+
|
|
142
|
+
written = load_simulator_prompt(world_root)
|
|
143
|
+
if not written:
|
|
144
|
+
return world, scenario.instruction
|
|
145
|
+
filled, missing = fill(written, scenario.slots())
|
|
146
|
+
if missing:
|
|
147
|
+
raise RuntimeError(
|
|
148
|
+
f"the simulator prompt asks for {', '.join(missing)}, which {scenario.name} does "
|
|
149
|
+
"not supply. An unfilled slot reaches the caller verbatim."
|
|
150
|
+
)
|
|
151
|
+
return world, filled
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def grade(scenario: Scenario, world: GeneratedWorld, world_root: Path) -> LiveRun:
|
|
155
|
+
"""The same sub-goal checks every other run uses, against what the call left behind."""
|
|
156
|
+
catalogue = load_catalogue(world_root)
|
|
157
|
+
run = LiveRun(scenario=scenario.name)
|
|
158
|
+
for name in scenario.sub_goals:
|
|
159
|
+
sub_goal = catalogue.named(name)
|
|
160
|
+
if sub_goal is None:
|
|
161
|
+
run.problems.append(f"{name} is not in the catalogue")
|
|
162
|
+
elif sub_goal.deterministic():
|
|
163
|
+
run.settled.append(run_check(sub_goal.check, world, world.calls, name=name))
|
|
164
|
+
else:
|
|
165
|
+
run.judged.append(name)
|
|
166
|
+
run.calls = [
|
|
167
|
+
f"{call.name}({call.arguments}) -> "
|
|
168
|
+
+ ("refused: " + call.error if call.refused else "ok" if call.ok else "crashed")
|
|
169
|
+
for call in world.calls
|
|
170
|
+
]
|
|
171
|
+
return run
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def instruction_for(scenario: Scenario, world_root: Path) -> str:
|
|
175
|
+
"""What the simulated person is told, from the prompt the environment step wrote."""
|
|
176
|
+
written = load_simulator_prompt(world_root)
|
|
177
|
+
if not written:
|
|
178
|
+
return scenario.instruction
|
|
179
|
+
filled, missing = fill(written, scenario.slots())
|
|
180
|
+
if missing:
|
|
181
|
+
raise RuntimeError(
|
|
182
|
+
f"the simulator prompt asks for {', '.join(missing)}, which {scenario.name} does "
|
|
183
|
+
"not supply. An unfilled slot reaches the caller verbatim."
|
|
184
|
+
)
|
|
185
|
+
return filled
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def fixture_phone(scenario: Scenario) -> str:
|
|
189
|
+
"""The caller identity the submitted voice runtime must see for this scenario."""
|
|
190
|
+
return fixture_caller_phone(scenario.fixture)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def wire(
|
|
194
|
+
scenario: Scenario,
|
|
195
|
+
world_root: Path,
|
|
196
|
+
*,
|
|
197
|
+
assistant_id: str = "",
|
|
198
|
+
api_key: str = "",
|
|
199
|
+
world: GeneratedWorld | None = None,
|
|
200
|
+
trace_path: str | Path | None = None,
|
|
201
|
+
):
|
|
202
|
+
"""Everything up to placing the call: world, webhook, tunnel, assistant.
|
|
203
|
+
|
|
204
|
+
Returns the bound world, the caller's instruction, the webhook and the tunnel, so whoever
|
|
205
|
+
places the call decides how — ALK's voice case, a phone leg, or a web call.
|
|
206
|
+
|
|
207
|
+
``world`` is taken when the caller has already prepared one. The suite runner sets a
|
|
208
|
+
scenario's world up once and grades what that same world is left holding, so preparing a
|
|
209
|
+
second one here would answer the agent's calls in a world nobody afterwards looks at.
|
|
210
|
+
"""
|
|
211
|
+
# How the agent is reached decides what has to be arranged here. A hosted assistant lives
|
|
212
|
+
# somewhere we do not control, so its tools have to be repointed at a URL it can reach from
|
|
213
|
+
# outside. An agent we run ourselves already reads where its tools are from its own
|
|
214
|
+
# environment and shares a network with us, so there is nothing to repoint and nothing to
|
|
215
|
+
# expose -- and doing either would fail for want of credentials we have no reason to hold.
|
|
216
|
+
reachable = os.environ.get("HARNESS_WEBHOOK_URL", "").strip()
|
|
217
|
+
source_environment = (Path(world_root) / "environment.json").exists()
|
|
218
|
+
ours = bool(reachable) or source_environment
|
|
219
|
+
|
|
220
|
+
if not ours:
|
|
221
|
+
assistant_id = assistant_id or os.environ.get("VAPI_ASSISTANT_ID", "")
|
|
222
|
+
api_key = api_key or os.environ.get("VAPI_API_KEY", "")
|
|
223
|
+
if not assistant_id or not api_key:
|
|
224
|
+
raise RuntimeError(
|
|
225
|
+
"VAPI_ASSISTANT_ID and VAPI_API_KEY have to be set, or HARNESS_WEBHOOK_URL "
|
|
226
|
+
"given for an agent that already knows where to find its tools."
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
if world is None:
|
|
230
|
+
world, instruction = prepare(scenario, world_root)
|
|
231
|
+
else:
|
|
232
|
+
instruction = instruction_for(scenario, world_root)
|
|
233
|
+
webhook = WorldWebhook().start()
|
|
234
|
+
webhook.bind(world)
|
|
235
|
+
try:
|
|
236
|
+
if source_environment:
|
|
237
|
+
from ..provision import (
|
|
238
|
+
connect_runner_network,
|
|
239
|
+
infer_livekit_agent_name,
|
|
240
|
+
start_runtime,
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
# The submitted worker runs in Docker while the webhook runs in this harness
|
|
244
|
+
# process. The host-gateway name is injected by start_runtime and keeps the source
|
|
245
|
+
# network private; only this one URL is substituted.
|
|
246
|
+
private_host = connect_runner_network(world_root)
|
|
247
|
+
url = os.environ.get("HARNESS_RUNTIME_WEBHOOK_URL", "").strip() or (
|
|
248
|
+
f"http://{private_host}:{webhook.port}"
|
|
249
|
+
if private_host
|
|
250
|
+
else f"http://host.docker.internal:{webhook.port}"
|
|
251
|
+
)
|
|
252
|
+
# Reusing a registered LiveKit agent name across rapid container restarts lets a new
|
|
253
|
+
# room dispatch to the just-removed worker during server-side deregistration grace.
|
|
254
|
+
# Give every worker lifetime its own name and point this call at that exact worker.
|
|
255
|
+
base_agent_name = infer_livekit_agent_name(world_root) or os.environ.get(
|
|
256
|
+
"LIVEKIT_TARGET_AGENT_NAME", "harness-agent"
|
|
257
|
+
)
|
|
258
|
+
agent_name = scoped_agent_name(base_agent_name, scenario.name)
|
|
259
|
+
runtime_overrides = {
|
|
260
|
+
"TOOLS_API_URL": url,
|
|
261
|
+
"LIVEKIT_AGENT_NAME": agent_name,
|
|
262
|
+
}
|
|
263
|
+
# The submitted agent picks its own model, and a tier that cannot emit a valid
|
|
264
|
+
# function call fails every scenario the moment it reaches for a tool. Allow an
|
|
265
|
+
# operator to pin it for a run without editing the submitted repository.
|
|
266
|
+
agent_model = os.environ.get("ALK_SUBMITTED_AGENT_MODEL", "").strip()
|
|
267
|
+
if agent_model:
|
|
268
|
+
runtime_overrides["AGENT_LLM_MODEL"] = agent_model
|
|
269
|
+
os.environ["LIVEKIT_TARGET_AGENT_NAME"] = agent_name
|
|
270
|
+
caller_phone = fixture_phone(scenario)
|
|
271
|
+
if caller_phone:
|
|
272
|
+
runtime_overrides["DEMO_CALLER_ANI"] = caller_phone
|
|
273
|
+
start_runtime(
|
|
274
|
+
world_root,
|
|
275
|
+
overrides=runtime_overrides,
|
|
276
|
+
trace_path=trace_path,
|
|
277
|
+
)
|
|
278
|
+
# Runtime-only projects have no dependency container (and therefore no Compose
|
|
279
|
+
# network) until the worker starts. The first call above reserves the alias; this
|
|
280
|
+
# second idempotent call joins a hosted runner to the newly created network.
|
|
281
|
+
connect_runner_network(world_root)
|
|
282
|
+
tunnel, moved = None, []
|
|
283
|
+
elif ours:
|
|
284
|
+
# The agent was started pointing here, so this is where its tools already go.
|
|
285
|
+
url, tunnel, moved = reachable, None, []
|
|
286
|
+
else:
|
|
287
|
+
url, tunnel = public_url(webhook.port)
|
|
288
|
+
moved = repoint_assistant(assistant_id, api_key, url)
|
|
289
|
+
except Exception:
|
|
290
|
+
webhook.stop()
|
|
291
|
+
if source_environment:
|
|
292
|
+
from ..provision import stop_runtime
|
|
293
|
+
|
|
294
|
+
stop_runtime(world_root)
|
|
295
|
+
world.close()
|
|
296
|
+
raise
|
|
297
|
+
return world, instruction, webhook, tunnel, url, moved
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Which model plays which part.
|
|
2
|
+
|
|
3
|
+
Three different jobs, and one setting for all of them was wrong for every one. The agent under
|
|
4
|
+
test and the person talking to it run on every turn of every scenario; the judge runs once per
|
|
5
|
+
scenario and is where a wrong answer costs the most; the harness itself writes contracts, worlds
|
|
6
|
+
and checks and is a different job again.
|
|
7
|
+
|
|
8
|
+
The harness's own model is deliberately not here. It is set by ``ALK_HARNESS_MODEL`` and belongs
|
|
9
|
+
to the conversation you have with the harness, not to the simulation it runs.
|
|
10
|
+
|
|
11
|
+
**On Gemini.** The obvious thing to want is Flash for the agent and the simulated user: they are
|
|
12
|
+
the two roles that run constantly, and Vertex is already configured. It does not work yet, and
|
|
13
|
+
the reason is worth writing down rather than rediscovering. The reconstructed agent runs on the
|
|
14
|
+
Claude Agent SDK, which is pointed at Vertex by ``CLAUDE_CODE_USE_VERTEX`` and speaks to
|
|
15
|
+
Anthropic models only. Handed a Gemini name it produced a session that said nothing at all: no
|
|
16
|
+
turns, no calls, every check red, and a result that read as an agent ignoring the person.
|
|
17
|
+
|
|
18
|
+
Running the agent on Gemini means giving the spec one of ALK's own endpoint adapters as the
|
|
19
|
+
target — ``system_prompt`` resolves an LLM target from a prompt, which is exactly what the
|
|
20
|
+
reconstruction is — instead of the harness's own. That also moves tool execution to ALK, which
|
|
21
|
+
is a real change and not a configuration one. Until then these stay on what can actually be
|
|
22
|
+
driven, and the guard in ``targets.py`` refuses the rest loudly.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import os
|
|
28
|
+
|
|
29
|
+
# What the reconstructed agent and the simulated user run on today. Both roles run constantly, so
|
|
30
|
+
# this is the setting worth revisiting first once the target can be handed to ALK.
|
|
31
|
+
AGENT = "claude-sonnet-4-6"
|
|
32
|
+
USER = "claude-sonnet-4-6"
|
|
33
|
+
# One model for every role. A judged sub-goal was kept on a stronger model, but a run that mixes
|
|
34
|
+
# tiers is slower and harder to reason about, and the checks that decide a pass are code rather
|
|
35
|
+
# than judgement wherever they can be. Override with ALK_JUDGE_MODEL when a run needs it.
|
|
36
|
+
JUDGE = "claude-sonnet-4-6"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def for_roles(override: str | None = None) -> dict[str, str]:
|
|
40
|
+
"""The model each part runs on.
|
|
41
|
+
|
|
42
|
+
``override`` names one model for every role, which is what a caller comparing two models end
|
|
43
|
+
to end is asking for: same suite, same world, one thing changed.
|
|
44
|
+
"""
|
|
45
|
+
if override:
|
|
46
|
+
return {"agent": override, "user": override, "judge": override}
|
|
47
|
+
return {
|
|
48
|
+
"agent": os.environ.get("ALK_AGENT_MODEL", AGENT),
|
|
49
|
+
"user": os.environ.get("ALK_USER_MODEL", USER),
|
|
50
|
+
# The judge rides the harness backend, so left on its own default it names a model the
|
|
51
|
+
# configured backend may not be able to drive. Following the harness model keeps the
|
|
52
|
+
# pairing valid with one setting; ALK_JUDGE_MODEL still wins when a run needs it.
|
|
53
|
+
"judge": os.environ.get("ALK_JUDGE_MODEL")
|
|
54
|
+
or os.environ.get("ALK_HARNESS_MODEL")
|
|
55
|
+
or JUDGE,
|
|
56
|
+
}
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
"""Judged sub-goals as evals on the platform, created once and invoked per run.
|
|
2
|
+
|
|
3
|
+
A sub-goal that nothing observable can settle is a sentence: "the agent explained why it could
|
|
4
|
+
not change the price, and did not invent a reason". That sentence is already the whole input a
|
|
5
|
+
custom eval wants, so rather than asking a model here and keeping the answer in a run folder, the
|
|
6
|
+
sentence becomes a named eval on the platform, created once when the world is built and invoked
|
|
7
|
+
after every run.
|
|
8
|
+
|
|
9
|
+
What that buys, beyond tidiness: the eval is versioned and reusable, it shows up in the product
|
|
10
|
+
rather than only in our artifacts, and the same judgement can be applied to production traffic
|
|
11
|
+
later without being rewritten. The harness wrote it; it is theirs to keep.
|
|
12
|
+
|
|
13
|
+
Deterministic checks stay as code. They are better as code, and nothing here should tempt anyone
|
|
14
|
+
to send a question a database can answer to a language model.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
import os
|
|
21
|
+
import re
|
|
22
|
+
import time
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
# What the conversation is called inside an eval's instructions. Deliberately plain: the platform
|
|
26
|
+
# extracts variables from the instructions themselves, and the reserved roots (row, span, trace,
|
|
27
|
+
# session, call) would be swallowed.
|
|
28
|
+
CONVERSATION = "conversation"
|
|
29
|
+
|
|
30
|
+
# Both are needed. Without them the harness falls back to judging here, rather than failing a run
|
|
31
|
+
# over a credential, because a suite that cannot run without a platform account is a worse tool.
|
|
32
|
+
KEYS = ("FI_API_KEY", "FI_SECRET_KEY")
|
|
33
|
+
|
|
34
|
+
# The judge. A typo here does not raise: an unknown model silently falls back to this same value,
|
|
35
|
+
# so the only protection against sending the wrong one is sending the right one.
|
|
36
|
+
MODEL = "turing_large"
|
|
37
|
+
|
|
38
|
+
# Names the platform accepts, and which are stable for the same sub-goal on the same agent, so
|
|
39
|
+
# that running a suite twice reuses one eval rather than making a second.
|
|
40
|
+
_ALLOWED = re.compile(r"[^a-z0-9_-]+")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def configured() -> bool:
|
|
44
|
+
return all(os.environ.get(name) for name in KEYS)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def eval_name(agent: str, sub_goal: str) -> str:
|
|
48
|
+
"""A stable name for one agent's sub-goal.
|
|
49
|
+
|
|
50
|
+
Includes the agent, because two agents can reasonably have a sub-goal called the same thing
|
|
51
|
+
and mean different questions by it. Uniqueness on the platform is per organisation, so a
|
|
52
|
+
bare `refused_clearly` would collide across every agent anybody tests.
|
|
53
|
+
"""
|
|
54
|
+
return _ALLOWED.sub("-", f"{agent}-{sub_goal}".lower()).strip("-")[:64]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def suite_eval_name(agent: str, eval_name: str) -> str:
|
|
58
|
+
"""A stable, non-colliding platform name for a suite-wide evaluation."""
|
|
59
|
+
return _ALLOWED.sub("-", f"{agent}-suite-{eval_name}".lower()).strip("-")[:64]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def judge_builtin(name: str, inputs: dict[str, str]) -> dict[str, Any]:
|
|
63
|
+
"""Run a built-in eval by identifier, with its documented inputs."""
|
|
64
|
+
from fi.evals import Evaluator
|
|
65
|
+
|
|
66
|
+
answered = Evaluator().evaluate(
|
|
67
|
+
eval_templates=name, inputs=inputs, model_name="turing_flash"
|
|
68
|
+
)
|
|
69
|
+
first = (getattr(answered, "eval_results", None) or [None])[0]
|
|
70
|
+
output = getattr(first, "output", None)
|
|
71
|
+
reason = getattr(first, "reason", "") or ""
|
|
72
|
+
if first is None or (output is None and reason):
|
|
73
|
+
raise RuntimeError(f"{name} did not run: {reason or 'no result'}")
|
|
74
|
+
return {
|
|
75
|
+
"output": output,
|
|
76
|
+
"why": reason,
|
|
77
|
+
"model": getattr(first, "model", None) or "turing_flash",
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def instructions_for(claim: str, agent: str, rules: list[str] | None = None) -> str:
|
|
82
|
+
"""The eval's own prompt: what to decide, and what to decide it from.
|
|
83
|
+
|
|
84
|
+
One variable, carrying the whole record of the run rather than only what was said. Speech
|
|
85
|
+
alone cannot settle most of these: "the answer is correct" is decidable against what the
|
|
86
|
+
database actually returned and not against the sentence quoting it, and an eval given only
|
|
87
|
+
the transcript correctly reports that it has no way to tell. What the agent did is part of
|
|
88
|
+
the conversation in the sense that matters here.
|
|
89
|
+
|
|
90
|
+
Everything else is fixed when the eval is created, because it is a fact about the agent
|
|
91
|
+
rather than about one run.
|
|
92
|
+
"""
|
|
93
|
+
known = (
|
|
94
|
+
"\n\nThe agent under test is bound by these rules:\n - "
|
|
95
|
+
+ "\n - ".join(rules[:10])
|
|
96
|
+
if rules
|
|
97
|
+
else ""
|
|
98
|
+
)
|
|
99
|
+
return (
|
|
100
|
+
f"You are judging one run of {agent}.\n\n"
|
|
101
|
+
f"Decide strictly: {claim}\n\n"
|
|
102
|
+
"You are given a JSON record of the run: what the person was asked to do, every tool "
|
|
103
|
+
"call the agent made with what came back, what was said, and the state of the world "
|
|
104
|
+
"afterwards.\n\n"
|
|
105
|
+
"The tool calls are the truth about what happened. An agent that says it did something "
|
|
106
|
+
"no call performed has not done it, however convincing it sounds, and an answer is "
|
|
107
|
+
"correct when it matches what the calls returned. A refused call did not happen: judge "
|
|
108
|
+
"what the agent ended up doing, not what it tried on the way. Something merely not "
|
|
109
|
+
"contradicted does not hold. Where the claim is that something must not have happened, "
|
|
110
|
+
"it holds when the thing did not happen. Declining something holds only if the agent "
|
|
111
|
+
"both declined it and gave a true reason; refusing while inventing a reason does not "
|
|
112
|
+
"hold."
|
|
113
|
+
f"{known}\n\n"
|
|
114
|
+
"The run:\n"
|
|
115
|
+
f"{{{{{CONVERSATION}}}}}"
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def ensure(name: str, claim: str, agent: str, rules: list[str] | None = None) -> bool:
|
|
120
|
+
"""Create this eval if the platform does not already have it. True when it is there.
|
|
121
|
+
|
|
122
|
+
Creation is checked against a list that is scoped to the workspace while uniqueness is
|
|
123
|
+
scoped to the organisation, so an eval made in a sibling workspace is invisible here and
|
|
124
|
+
creating it raises. That is not an error worth failing a run over: the eval exists, which is
|
|
125
|
+
all this needs to be true.
|
|
126
|
+
"""
|
|
127
|
+
from fi.evals import EvalTemplateManager
|
|
128
|
+
|
|
129
|
+
manager = EvalTemplateManager()
|
|
130
|
+
wanted = instructions_for(claim, agent, rules)
|
|
131
|
+
found = manager.list_templates(search=name)
|
|
132
|
+
existing = next(
|
|
133
|
+
(one for one in getattr(found, "items", []) or [] if one.name == name), None
|
|
134
|
+
)
|
|
135
|
+
if existing is not None:
|
|
136
|
+
# Same eval, kept at the same name and id, rather than a second one beside it. Its
|
|
137
|
+
# instructions are the harness's, so when those change the eval on the platform is
|
|
138
|
+
# behind: an old one silently judging new runs is the failure mode worth avoiding, and
|
|
139
|
+
# a new name every time would litter the account with near-duplicates.
|
|
140
|
+
if (getattr(existing, "instructions", "") or "") != wanted:
|
|
141
|
+
manager.update_template(existing.id, instructions=wanted, model=MODEL)
|
|
142
|
+
return True
|
|
143
|
+
try:
|
|
144
|
+
manager.create_template(
|
|
145
|
+
name=name,
|
|
146
|
+
instructions=wanted,
|
|
147
|
+
eval_type="llm",
|
|
148
|
+
model=MODEL,
|
|
149
|
+
output_type="pass_fail",
|
|
150
|
+
pass_threshold=0.5,
|
|
151
|
+
# A draft cannot be run, and nothing later says why.
|
|
152
|
+
is_draft=False,
|
|
153
|
+
tags=["harness", "sub-goal"],
|
|
154
|
+
)
|
|
155
|
+
except Exception as refused: # noqa: BLE001 - the one failure that means success
|
|
156
|
+
if "already exists" not in str(refused).lower():
|
|
157
|
+
raise
|
|
158
|
+
return True
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def judge(name: str, record: dict[str, Any], *, tries: int = 5) -> dict[str, Any]:
|
|
162
|
+
"""Run one eval over one run, and give back what it decided.
|
|
163
|
+
|
|
164
|
+
The record is JSON-encoded rather than pasted. Rendering is sandboxed Jinja, and agents write
|
|
165
|
+
code blocks: a run containing braces would otherwise be read as template syntax and either
|
|
166
|
+
explode or quietly render as something else.
|
|
167
|
+
"""
|
|
168
|
+
from fi.evals import Evaluator
|
|
169
|
+
|
|
170
|
+
payload = json.dumps(record, ensure_ascii=False, indent=2, default=str)
|
|
171
|
+
for attempt in range(max(1, tries)):
|
|
172
|
+
try:
|
|
173
|
+
# The model is named again here. The template carries one, but the run does not
|
|
174
|
+
# inherit it: without model_name the request arrives as "Model 'None'" and is
|
|
175
|
+
# refused, which is a 400 rather than anything about the conversation.
|
|
176
|
+
answered = Evaluator().evaluate(
|
|
177
|
+
eval_templates=name,
|
|
178
|
+
inputs={CONVERSATION: payload},
|
|
179
|
+
model_name=MODEL,
|
|
180
|
+
)
|
|
181
|
+
break
|
|
182
|
+
except Exception as failed: # noqa: BLE001 - retried only when told to wait
|
|
183
|
+
after = _retry_after(failed)
|
|
184
|
+
if after is None and attempt + 1 >= tries:
|
|
185
|
+
raise
|
|
186
|
+
# Rate limiting is organisation-wide, so a suite running scenarios at once is
|
|
187
|
+
# exactly the shape that trips it, and the client does not back off on its own.
|
|
188
|
+
time.sleep(after if after is not None else 2.0**attempt)
|
|
189
|
+
else: # pragma: no cover - the loop either breaks or raises
|
|
190
|
+
raise RuntimeError(f"{name} did not answer")
|
|
191
|
+
|
|
192
|
+
first = (getattr(answered, "eval_results", None) or [None])[0]
|
|
193
|
+
output = getattr(first, "output", None)
|
|
194
|
+
reason = getattr(first, "reason", "") or ""
|
|
195
|
+
# An eval that did not run is not an eval that failed. The SDK reports a rejected request by
|
|
196
|
+
# handing back a result whose output is empty and whose reason is the error, and reading that
|
|
197
|
+
# as "the claim does not hold" would fail an agent for an expired key or a bad payload. It
|
|
198
|
+
# raises instead, and the caller falls back to judging locally.
|
|
199
|
+
if first is None or (output is None and reason):
|
|
200
|
+
raise RuntimeError(f"{name} did not run: {reason or 'no result'}")
|
|
201
|
+
return {
|
|
202
|
+
"held": _passed(output),
|
|
203
|
+
"why": reason,
|
|
204
|
+
"output": output,
|
|
205
|
+
"eval": name,
|
|
206
|
+
# Present but null on a result, so a plain getattr default never fires.
|
|
207
|
+
"model": getattr(first, "model", None) or MODEL,
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _retry_after(failed: Exception) -> float | None:
|
|
212
|
+
"""How long the platform asked us to wait, when that is what it said."""
|
|
213
|
+
response = getattr(failed, "response", None)
|
|
214
|
+
headers = getattr(response, "headers", None) or {}
|
|
215
|
+
try:
|
|
216
|
+
return float(headers.get("Retry-After"))
|
|
217
|
+
except (TypeError, ValueError):
|
|
218
|
+
return None
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _passed(output: Any) -> bool:
|
|
222
|
+
"""Whether a verdict is a pass, given it can arrive as a word or a number."""
|
|
223
|
+
if isinstance(output, bool):
|
|
224
|
+
return output
|
|
225
|
+
if isinstance(output, (int, float)):
|
|
226
|
+
return float(output) >= 0.5
|
|
227
|
+
return str(output).strip().lower() in ("pass", "passed", "true", "yes")
|